"""
@FileName：test.py
@Description：

@Author：HeYiQing
@Time：2023/11/27 20:30
"""
import openpyxl
import requests
import re
from lxml import etree
from bs4 import BeautifulSoup

# title article
if __name__ == '__main__':
    # UA伪装
    headers = {
        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36'
    }

    url = 'http://www.dili360.com/article/p5474acf05c91d67.htm'
    page_text = requests.get(url=url, headers=headers).text

    tree = etree.HTML(page_text)
    page_xpath = tree.xpath('//div[contains(@class, "common-article-page")]')[0]
    page = etree.tostring(page_xpath, pretty_print=True).decode()
    # print(page)

    pattern = re.compile(r'<article>(.*?)</article>', re.DOTALL)
    match = pattern.search(page_text)
    group = match.group(1)
    re_compile = re.compile(r'<h1>(.*?)<br>', re.DOTALL)
    h1 = re_compile.findall(group)
    print(h1)



